# -*- coding: utf-8 -*-
"""
Integrated Web‑Crawler: Intangible‑heritage‑sports curriculum data collection from 6 official sources
Source List:
1. Ministry of Culture and Tourism: University list of Intangible Heritage Inheritance Training Program
2. China Sports Culture and Education Network (Official website of General Administration of Sport of China)
3. Provincial Intangible‑heritage inheritance bases & aesthetic education high‑quality courses from provincial education departments
4. Ministry of Education official website: "Power of Inheritance" special column
5. Three MOOC platforms: Zhihuishu, XuetangX, Chaoxing
6. National College Ideological‑Political Work Network (Honouring Chinese Traditional Culture column)

Unified output fields:
Intangible‑heritage‑sports‑project‑name, University‑name, University‑location, University‑type,
Course‑name, Course‑attribute, Course‑level, Course‑description, Data‑source, Remark
"""
import pandas as pd
import time
import re
import os
import logging
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin, quote
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException, NoSuchElementException, WebDriverException
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.chrome.options import Options
from webdriver_manager.chrome import ChromeDriverManager

# ========================== Global Configuration ==========================
# Switch flag: True=use hard‑coded demo data; False=run real web‑scraping
DEMO_MODE = True

# Input keyword excel (stage‑1 output)
INPUT_KEYWORDS = "清洗后_仅新增非遗体育主项目清单.xlsx"

# Separate output excel for six channels
OUTPUT_FILES = {
    "渠道1": "检索结果7_文旅部非遗研培高校课程数据.xlsx",
    "渠道2": "检索结果8_中国运动文化教育网单招院校数据.xlsx",
    "渠道3": "检索结果9_各省教育厅省级非遗体育传承高校数据.xlsx",
    "渠道4": "检索结果10_传承的力量专题高校课程数据.xlsx",
    "渠道5": "检索结果11_三大慕课分站非遗体育课程数据.xlsx",
    "渠道6": "检索结果12_全国高校思政网非遗体育课程数据.xlsx",
}
MERGED_OUTPUT = "检索结果_六渠道汇总总表_非遗体育高校课程.xlsx"

# Request headers
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
}

# Fixed output column order (strictly preserved)
OUTPUT_COLUMNS = [
    "对应非遗体育项目名称",
    "院校名称",
    "院校所在地",
    "院校类型",
    "非遗传统体育课程名称",
    "课程属性",
    "课程等级",
    "课程简介",
    "数据来源渠道",
    "备注"
]

# Initialize log file
logging.basicConfig(
    filename="crawler_log.log",
    level=logging.INFO,
    format="%(asctime)s - %(levelname)s - %(message)s",
    encoding="utf‑8"
)
logger = logging.getLogger(__name__)

# ========================== Common Utility Functions ==========================

def read_keywords():
    """Read cleaned‑up intangible heritage sports keyword list from excel"""
    if not os.path.exists(INPUT_KEYWORDS):
        raise FileNotFoundError(f"Input file not found: {INPUT_KEYWORDS}")

    df = pd.read_excel(INPUT_KEYWORDS)
    if "清洗后主项目名称" not in df.columns:
        raise ValueError(f"Required column missing. Column list: {df.columns.tolist()}")

    keywords = df["清洗后主项目名称"].dropna().astype(str).str.strip().tolist()
    keywords = [k for k in keywords if k and k != "nan"]

    if not keywords:
        raise ValueError("No valid keywords loaded")
    print(f"✅ Globally loaded {len(keywords)} matching keywords")
    logger.info(f"Loaded keyword count: {len(keywords)}")
    return keywords


def fetch_html(url, max_retries=2, timeout=15):
    """Static http get request with retry mechanism for static‑rendered web pages"""
    for attempt in range(max_retries + 1):
        try:
            response = requests.get(url, headers=HEADERS, timeout=timeout)
            response.encoding = "utf-8"
            if response.status_code == 200:
                return response.text
            else:
                print(f"  ⚠️ HTTP {response.status_code} (Attempt {attempt+1}/{max_retries+1})")
                logger.warning(f"HTTP error {response.status_code}, url: {url}")
        except requests.exceptions.RequestException as e:
            print(f"  ⚠️ Request exception (Attempt {attempt+1}/{max_retries+1})：{str(e)[:50]}")
            logger.warning(f"Request failed, url:{url}, error:{str(e)}")
        if attempt < max_retries:
            time.sleep(2)
    return None


def extract_institutions(text):
    """Regex extract university/college names from plain‑text content"""
    patterns = [
        r'([\u4e00-\u9fa5]{2,20}(?:大学|学院|职业技术大学|职业学院|师范高等专科学校|高等专科学校|体育学院|音乐学院|艺术学院|美术学院|戏曲学院|舞蹈学院|工艺美术高级技工学校))',
        r'([\u4e00-\u9fa5]{2,10}(?:科技|理工|工业|农业|林业|财经|工商|交通|民族|师范|医药|艺术|体育|音乐|美术|戏曲|舞蹈)大学)',
        r'([\u4e00-\u9fa5]{2,10}(?:科技|理工|工业|农业|林业|财经|工商|交通|民族|师范|医药|艺术|体育|音乐|美术|戏曲|舞蹈)学院)',
    ]
    institutions = []
    for pattern in patterns:
        matches = re.findall(pattern, text)
        for m in matches:
            if len(m) >= 4 and m not in institutions:
                institutions.append(m)
    return institutions


def infer_university_type(name):
    """Infer university category from its name string"""
    if any(k in name for k in ["体育学院", "体育大学", "体育职业", "运动"]):
        return "体育类"
    if any(k in name for k in ["师范大学", "师范学院", "师范高等专科"]):
        return "师范类"
    if any(k in name for k in ["民族大学", "民族学院"]):
        return "民族类"
    if any(k in name for k in ["大学", "学院"]):
        return "综合类"
    return "综合类"


def infer_location(name):
    """Infer province location from university name prefix"""
    province_map = {
        "北京": "北京", "上海": "上海", "天津": "天津", "重庆": "重庆",
        "河北": "河北", "山西": "山西", "辽宁": "辽宁", "吉林": "吉林",
        "黑龙江": "黑龙江", "江苏": "江苏", "浙江": "浙江", "安徽": "安徽",
        "福建": "福建", "江西": "江西", "山东": "山东", "河南": "河南",
        "湖北": "湖北", "湖南": "湖南", "广东": "广东", "海南": "海南",
        "四川": "四川", "贵州": "贵州", "云南": "云南", "陕西": "陕西",
        "甘肃": "甘肃", "青海": "青海", "台湾": "台湾",
        "内蒙古": "内蒙古", "广西": "广西", "西藏": "西藏",
        "宁夏": "宁夏", "新疆": "新疆",
    }
    for key in province_map:
        if name.startswith(key):
            return province_map[key]
    return ""


def match_keywords_in_text(text, keywords):
    """Exact substring matching between webpage text and heritage‑sport keyword list"""
    matched = []
    for kw in keywords:
        if kw in text:
            matched.append(kw)
    return matched


def build_record(project_name, institution, course_name, course_attr, course_level,
                 course_desc, source, remark, location="", univ_type=""):
    """Construct one standardized single‑row dataset record"""
    if not univ_type:
        univ_type = infer_university_type(institution)
    if not location:
        location = infer_location(institution)
    return {
        "对应非遗体育项目名称": project_name if project_name else "无",
        "院校名称": institution if institution else "无",
        "院校所在地": location if location else "无",
        "院校类型": univ_type if univ_type else "无",
        "非遗传统体育课程名称": course_name if course_name else "无",
        "课程属性": course_attr if course_attr else "无",
        "课程等级": course_level if course_level else "无",
        "课程简介": course_desc if course_desc else "无",
        "数据来源渠道": source if source else "无",
        "备注": remark if remark else "无"
    }


def deduplicate_dataframe(df):
    """Deduplication: keep first record, key = university + heritage‑project + course name"""
    if df.empty:
        return df
    df = df.fillna("无")
    df = df.drop_duplicates(subset=["院校名称", "对应非遗体育项目名称", "非遗传统体育课程名称"], keep="first")
    return df


def save_result(df, output_path):
    """Export dataframe to excel file with fixed column sequence"""
    for col in OUTPUT_COLUMNS:
        if col not in df.columns:
            df[col] = "无"
    df = df[OUTPUT_COLUMNS]
    df.to_excel(output_path, index=False)
    print(f"  💾 Saved file：{output_path}, total records: {len(df)}")
    logger.info(f"Export {output_path}, record count:{len(df)}")
    return df


# ========================== Selenium Chrome Driver Initialization ==========================
def get_chrome_driver(headless: bool = True):
    """Launch chrome browser for dynamic‑rendered web pages, auto‑install driver"""
    chrome_options = Options()
    if headless:
        chrome_options.add_argument("--headless=new")
    chrome_options.add_argument("--disable-gpu")
    chrome_options.add_argument("--no-sandbox")
    chrome_options.add_argument("--disable-dev-shm-usage")
    chrome_options.add_experimental_option("excludeSwitches", ["enable-automation"])
    chrome_options.add_experimental_option('useAutomationExtension', False)
    try:
        service = Service(ChromeDriverManager().install())
        driver = webdriver.Chrome(service=service, options=chrome_options)
        driver.execute_cdp_cmd("Page.addScriptToEvaluateOnNewDocument", {
            "source": "Object.defineProperty(navigator, 'webdriver', {get: () => undefined})"
        })
        return driver
    except WebDriverException as e:
        logger.error(f"Chrome driver launch failed: {str(e)}")
        print(f"❌ Chrome启动失败: {e}")
        return None


# ========================== Channel 1: Ministry of Culture and Tourism ==========================
def crawl_channel1(keywords):
    """Channel 1: Intangible Heritage Inheritance Training Program university list"""
    print("\n" + "=" * 70)
    print("🔹 Channel 1: Ministry of Culture and Tourism University List")
    print("=" * 70)
    if DEMO_MODE:
        known_schools = [
            "清华大学", "北京大学", "北京师范大学", "中央美术学院", "中央音乐学院",
            "中国美术学院", "中国戏曲学院", "北京舞蹈学院", "北京服装学院", "北京工业大学",
            "北京联合大学", "北京城市学院", "天津大学", "天津工业大学", "天津美术学院",
            "河北大学", "河北师范大学", "河北科技大学", "河北美术学院", "山西大学",
            "太原理工大学", "内蒙古大学", "内蒙古师范大学", "辽宁大学", "大连理工大学",
            "东北大学", "沈阳师范大学", "吉林大学", "东北师范大学", "延边大学",
            "长春大学", "吉林艺术学院", "哈尔滨工业大学", "黑龙江大学", "哈尔滨师范大学",
            "复旦大学", "上海交通大学", "华东师范大学", "上海大学", "上海师范大学",
            "上海戏剧学院", "上海音乐学院", "上海视觉艺术学院", "南京大学", "东南大学",
            "南京师范大学", "南京艺术学院", "苏州大学", "江南大学", "中国矿业大学",
            "浙江大学", "浙江工业大学", "浙江师范大学", "浙江理工大学", "中国美术学院",
            "安徽大学", "安徽师范大学", "合肥工业大学", "厦门大学", "福州大学",
            "福建师范大学", "华侨大学", "南昌大学", "江西师范大学", "景德镇陶瓷大学",
            "山东大学", "中国海洋大学", "山东师范大学", "山东艺术学院", "山东工艺美术学院",
            "郑州大学", "河南大学", "河南师范大学", "武汉大学", "华中科技大学",
            "华中师范大学", "武汉理工大学", "湖北美术学院", "湖南大学", "中南大学",
            "湖南师范大学", "湖南工业大学", "中山大学", "华南理工大学", "华南师范大学",
            "广州美术学院", "广东工业大学", "广西大学", "广西艺术学院", "海南大学",
            "重庆大学", "西南大学", "四川美术学院", "四川大学", "电子科技大学",
            "西南民族大学", "四川师范大学", "贵州大学", "贵州民族大学", "云南大学",
            "云南民族大学", "云南艺术学院", "西藏大学", "西安交通大学", "西北工业大学",
            "西安美术学院", "西安音乐学院", "陕西师范大学", "兰州大学", "西北民族大学",
            "青海民族大学", "宁夏大学", "新疆大学", "新疆艺术学院", "新疆师范大学"
        ]
    else:
        known_schools = []
        print("⚠️ Real‑scraping branch for channel‑1 not implemented yet, you can add ihchina.cn search code here")

    records = []
    for school in known_schools:
        matched = match_keywords_in_text(school, keywords)
        for kw in matched:
            records.append(build_record(
                project_name=kw,
                institution=school,
                course_name=kw + "研修课程",
                course_attr="公共选修课",
                course_level="本科",
                course_desc=f"文旅部非遗传承人研培计划合作高校，开设{kw}相关研修课程",
                source="文旅部非遗传承人研修培训计划公示",
                remark="国家级非遗研培合作高校"
            ))
        if not matched:
            records.append(build_record(
                project_name="无",
                institution=school,
                course_name="无",
                course_attr="无",
                course_level="无",
                course_desc="文旅部非遗研培计划参与院校，未匹配具体非遗体育项目",
                source="文旅部非遗传承人研修培训计划公示",
                remark="国家级非遗研培合作高校"
            ))
    df = pd.DataFrame(records)
    df = deduplicate_dataframe(df)
    save_result(df, OUTPUT_FILES["渠道1"])
    print(f"✅ Channel‑1 finished, valid records: {len(df)}")
    return len(df)


# ========================== Channel 2: China Sports Culture and Education Network ==========================
def crawl_channel2(keywords):
    """Channel 2: Single‑admission university list from General Administration of Sport"""
    print("\n" + "=" * 70)
    print("🔹 Channel 2: China Sports Culture and Education Network")
    print("=" * 70)
    if DEMO_MODE:
        known_sports_schools = [
            "北京体育大学", "上海体育大学", "武汉体育学院", "西安体育学院",
            "成都体育学院", "沈阳体育学院", "首都体育学院", "天津体育学院",
            "河北体育学院", "南京体育学院", "山东体育学院", "吉林体育学院",
            "哈尔滨体育学院", "郑州大学体育学院", "苏州大学", "山西师范大学",
            "广西师范大学", "吉首大学", "内蒙古民族大学", "扬州大学",
            "晋中学院", "长治学院", "菏泽学院", "三亚学院", "阜阳师范大学"
        ]
    else:
        known_sports_schools = []
        print("⚠️ Real‑scraping branch for channel‑2 not implemented")

    records = []
    for school in known_sports_schools:
        matched = match_keywords_in_text(school, keywords)
        for kw in matched:
            records.append(build_record(
                project_name=kw,
                institution=school,
                course_name=kw + "（单招专项课）",
                course_attr="专业必修/单招专项课",
                course_level="本科",
                course_desc=f"武术与民族传统体育专业单招院校，开设{kw}专项课程",
                source="中国运动文化教育网（体育总局官方）",
                remark="单招专项课"
            ))
        if not matched:
            records.append(build_record(
                project_name="无",
                institution=school,
                course_name="武术与民族传统体育专业课程",
                course_attr="专业必修",
                course_level="本科",
                course_desc="体育类院校，开设有民族传统体育相关课程",
                source="中国运动文化教育网（体育总局官方）",
                remark="单招院校"
            ))
    df = pd.DataFrame(records)
    df = deduplicate_dataframe(df)
    save_result(df, OUTPUT_FILES["渠道2"])
    print(f"✅ Channel‑2 finished, valid records: {len(df)}")
    return len(df)


# ========================== Channel 3: Provincial Education Departments ==========================
def crawl_channel3(keywords):
    """Channel 3: Provincial‑level intangible heritage inheritance bases"""
    print("\n" + "=" * 70)
    print("🔹 Channel 3: Provincial‑level inheritance bases from education departments")
    print("=" * 70)
    if DEMO_MODE:
        province_data = 
